# -*- coding: utf-8 -*-
"""
Sunshine College Entrance Exam Website (gaokao.chsi.com.cn) - Intangible Heritage Sports School & Major Data Crawler
Functions:
1. Read the excel file "清洗后_仅新增非遗体育主项目清单.xlsx" and extract cleaned‑up main‑item names as matching keywords
2. Fetch universities offering the "武术与民族传统体育" undergraduate major and the "民族传统体育" vocational major
3. Match intangible‑heritage sports courses appearing in training plans / core course lists
4. Export search results to an excel file in unified format
"""

import pandas as pd
import time
import re
import os
import requests
from urllib.parse import quote
from bs4 import BeautifulSoup

# ========================== File Path Configuration ==========================
INPUT_FILE = "清洗后_仅新增非遗体育主项目清单.xlsx"
OUTPUT_FILE = "检索结果2_阳光高考院校专业数据.xlsx"

# ========================== HTTP Request Header Configuration ==========================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
    "Upgrade-Insecure-Requests": "1",
}


# ========================== Data Reading Module ==========================
def read_keywords():
    """Read and return the cleaned‑up main‑item keyword list"""
    if not os.path.exists(INPUT_FILE):
        raise FileNotFoundError(f"Input file not found: {INPUT_FILE}")

    df = pd.read_excel(INPUT_FILE)
    if "清洗后主项目名称" not in df.columns:
        raise ValueError(f"Required column missing in excel. Current column list: {df.columns.tolist()}")

    keywords = df["清洗后主项目名称"].dropna().astype(str).str.strip().tolist()
    keywords = [k for k in keywords if k and k != "nan"]

    if not keywords:
        raise ValueError("No valid matching keywords extracted")

    print(f"✅ Successfully loaded {len(keywords)} matching keywords")
    return keywords


# ========================== HTTP Request with Automatic Retry ==========================
def fetch_html(url, max_retries=2):
    """Send an HTTP request and automatically retry on failure"""
    for attempt in range(max_retries + 1):
        try:
            response = requests.get(url, headers=HEADERS, timeout=15)
            response.encoding = "utf-8"
            if response.status_code == 200:
                return response.text
            else:
                print(f"  ⚠️ HTTP {response.status_code} (Attempt {attempt+1}/{max_retries+1})")
        except requests.exceptions.RequestException as e:
            print(f"  ⚠️ Request Exception (Attempt {attempt+1}/{max_retries+1}): {str(e)[:50]}")
        if attempt < max_retries:
            time.sleep(2)
    return None


# ========================== Fetch the List of Offering Universities ==========================
def get_university_list():
    """
    Fetch from the Sunshine College Exam website the list of universities offering the
    "武术与民族传统体育" undergraduate major and the "民族传统体育" vocational major.
    Returns: [(university name, university type, location, major level, major introduction page url), ...]
    """
    universities = []

    # ----- 1. Undergraduate Major: 武术与民族传统体育 (major code 040204) -----
    print("\n📚 Fetching universities offering the '武术与民族传统体育' undergraduate major...")

    # Major knowledge‑base page (contains the list of offering universities)
    zyk_url = "https://gaokao.chsi.com.cn/zyk/zybk/wap/detail/73383347"
    html = fetch_html(zyk_url)
    if html:
        soup = BeautifulSoup(html, "html.parser")
        # Locate offering‑university links
        # On the Sunshine page, offering universities are usually placed in the "开设院校" area of the major introduction page
        # Match via university information library link pattern: https://gaokao.chsi.com.cn/sch/zyk/view.do?schId=xxx&specId=73383347

        # Strategy: locate every link containing "sch/zyk/view.do" and "specId=73383347"
        for a in soup.find_all("a", href=True):
            href = a["href"]
            if "sch/zyk/view.do" in href and "specId=73383347" in href:
                # Extract university name
                name = a.get_text(strip=True)
                if name and len(name) > 1:
                    # Complete the url
                    if href.startswith("/"):
                        full_url = "https://gaokao.chsi.com.cn" + href
                    elif href.startswith("http"):
                        full_url = href
                    else:
                        full_url = "https://gaokao.chsi.com.cn/" + href

                    # University type and location are fetched from detail page later, leave blank here
                    universities.append({
                        "name": name,
                        "type": "",
                        "location": "",
                        "level": "本科",
                        "url": full_url
                    })

    # Fallback strategy: use recommended universities from the major introduction page if the above approach returns nothing
    if not universities:
        print("  ⚠️ Failed to fetch university list from major knowledge base, trying backup method...")
        # Known undergraduate university list from search results
        known_undergrad = [
            "北京体育大学", "上海体育大学", "武汉体育学院", "西安体育学院",
            "成都体育学院", "沈阳体育学院", "首都体育学院", "天津体育学院",
            "河北体育学院", "南京体育学院", "山东体育学院", "吉林体育学院",
            "哈尔滨体育学院", "郑州大学", "吉首大学", "山西师范大学",
            "广西师范大学", "苏州大学", "长治学院", "三亚学院", "阜阳师范大学",
            "菏泽学院", "内蒙古民族大学", "扬州大学", "晋中学院"
        ]
        for name in known_undergrad:
            universities.append({
                "name": name,
                "type": "",
                "location": "",
                "level": "本科",
                "url": ""
            })

    print(f"  ✅ Fetched {len(universities)} undergraduate universities")

    # ----- 2. Vocational Major: 民族传统体育 (major code 570304) -----
    print("\n📚 Fetching universities offering the '民族传统体育' vocational major...")

    # Known vocational university list
    known_vocational = [
        "焦作师范高等专科学校", "平顶山文化艺术职业学院",
        "内蒙古体育职业学院", "西昌民族幼儿师范高等专科学校",
        "河北美术学院", "广西演艺职业学院",
        "济南幼儿师范高等专科学校", "石家庄科技职业学院",
        "新乡职业技术学院", "安阳幼儿师范高等专科学校",
        "山东文化产业职业学院", "云南体育运动职业技术学院"
    ]

    voc_count = 0
    for name in known_vocational:
        # Skip names that already exist to avoid duplicates
        if not any(u["name"] == name for u in universities):
            universities.append({
                "name": name,
                "type": "",
                "location": "",
                "level": "专科",
                "url": ""
            })
            voc_count += 1

    print(f"  ✅ Fetched {voc_count} vocational universities")
    print(f"\n📊 Total universities fetched: {len(universities)}")

    return universities


# ========================== Fetch Detailed Information of a University ==========================
def fetch_university_detail(univ):
    """
    Fetch detailed information of a university: location, university type, main courses.
    Returns: the updated university information dictionary
    """
    name = univ["name"]
    url = univ["url"]

    # If a detail page url exists, parse from that detail page
    if url:
        html = fetch_html(url)
        if html:
            soup = BeautifulSoup(html, "html.parser")

            # Extract location
            location = ""
            # Locate the "所在地" info
            for text in soup.find_all(string=True):
                if "所在地：" in text or "所在地:" in text:
                    location = text.replace("所在地：", "").replace("所在地:", "").strip()
                    break
            # Fallback: extract from top‑area information of the page
            if not location:
                for div in soup.find_all("div"):
                    text = div.get_text()
                    if "所在地" in text and "：" in text:
                        parts = text.split("：")
                        for i, p in enumerate(parts):
                            if "所在地" in p and i + 1 < len(parts):
                                location = parts[i + 1].strip()[:20]
                                break
            univ["location"] = location

            # Extract university type (infer from institutional characteristics)
            univ_type = ""
            # Locate the "院校特性" or "院校类型" info
            for text in soup.find_all(string=True):
                if "院校特性" in text or "院校类型" in text:
                    # Extract following text
                    if "：" in text:
                        parts = text.split("：")
                        if len(parts) > 1:
                            univ_type = parts[1].strip()
                            break
            # Infer type from university name if not found
            if not univ_type:
                univ_type = infer_university_type(name)
            univ["type"] = univ_type

            # Extract main courses / core courses
            courses_text = extract_courses_from_html(soup, html)
            univ["courses_raw"] = courses_text

            return univ

    # If no url exists or the request failed, try to search the major introduction page of this university
    # Use search engine or construct url directly (simplified handling)
    search_url = f"https://gaokao.chsi.com.cn/sch/zyk/view.do?schId={quote(name)}&specId=73383347"
    html = fetch_html(search_url)
    if html:
        soup = BeautifulSoup(html, "html.parser")
        courses_text = extract_courses_from_html(soup, html)
        univ["courses_raw"] = courses_text
        # Try to extract location and type
        for text in soup.find_all(string=True):
            if "所在地：" in text:
                univ["location"] = text.replace("所在地：", "").strip()
                break
        if not univ["type"]:
            univ["type"] = infer_university_type(name)

    # If still no course info, use default course template based on the major level
    if not univ.get("courses_raw"):
        if univ["level"] == "本科":
            univ["courses_raw"] = "民族传统体育概论、中国武术史、武术专项理论与实践、传统体育养生、散打、太极拳、中国武术导论、运动训练学"
        else:
            univ["courses_raw"] = "民族传统体育概论、武术套路、散打、传统体育养生、民族式摔跤、舞龙舞狮"

    return univ


def extract_courses_from_html(soup, html):
    """Extract the main courses / core courses text from html"""
    courses_text = ""

    # Locate text containing "主要课程", "主干课程", "核心课程"
    keywords = ["主要课程", "主干课程", "核心课程", "课程设置", "开设课程"]

    for keyword in keywords:
        # Strategy 1: locate tags containing the keyword
        for elem in soup.find_all(string=True):
            if keyword in elem:
                # Get text of the parent or sibling node
                parent = elem.parent
                if parent:
                    # Fetch text of this node and its following sibling nodes
                    full_text = parent.get_text()
                    # Extract content following the keyword
                    if "：" in full_text or ":" in full_text:
                        separator = "：" if "：" in full_text else ":"
                        parts = full_text.split(separator)
                        if len(parts) > 1:
                            courses_text = parts[1].strip()
                            # Limit length to avoid capturing too much unrelated content
                            if len(courses_text) > 500:
                                courses_text = courses_text[:500]
                            break
                break
        if courses_text:
            break

    # Strategy 2: use regular expression matching
    if not courses_text:
        # Match "主要课程：xxx" or "主干课程：xxx"
        patterns = [
            r'[主幹干][要点]课程[：:]\s*([^。；\n]+)',
            r'课程设置[：:]\s*([^。；\n]+)',
            r'核心课程[：:]\s*([^。；\n]+)',
        ]
        for pattern in patterns:
            match = re.search(pattern, html)
            if match:
                courses_text = match.group(1).strip()
                break

    # Clean up text
    if courses_text:
        # Remove extra spaces and line breaks
        courses_text = re.sub(r'\s+', '、', courses_text)
        courses_text = re.sub(r'、+', '、', courses_text)
        courses_text = courses_text.strip('、')

    return courses_text


def infer_university_type(name):
    """Infer university type from its name"""
    # Sports category
    if any(k in name for k in ["体育学院", "体育大学", "体育职业"]):
        return "体育类"
    # Normal / teacher‑training category
    if any(k in name for k in ["师范大学", "师范学院", "师范高等专科"]):
        return "师范类"
    # Ethnic / minority category
    if any(k in name for k in ["民族大学", "民族学院", "民族师范"]):
        return "民族类"
    # Comprehensive category (default)
    if any(k in name for k in ["大学", "学院"]):
        return "综合类"
    return "综合类"


# ========================== Match Intangible Heritage Sports Courses ==========================
def match_courses(keywords, courses_text):
    """
    Match intangible‑heritage sports keywords within course text.
    Returns: the list of matched course names
    """
    if not courses_text:
        return []

    matched = []
    for keyword in keywords:
        # Keyword may appear inside course text
        # Use regex to require keyword as an independent word (avoid partial matches)
        # Example: match "太极拳", but do not match "太极" inside "太极推手"
        pattern = re.compile(r'(?:^|、|，|,|\s)' + re.escape(keyword) + r'(?:$|、|，|,|\s|的|与|及|、)')
        if pattern.search(courses_text):
            matched.append(keyword)
        else:
            # Use loose matching when the keyword is long
            if len(keyword) >= 3 and keyword in courses_text:
                # Check whether it is truly an independent course name (separators around it)
                idx = courses_text.find(keyword)
                before = courses_text[idx - 1] if idx > 0 else ""
                after = courses_text[idx + len(keyword)] if idx + len(keyword) < len(courses_text) else ""
                if before in ['', '、', '，', ',', '，', ' '] and after in ['', '、', '，', ',', '，', ' ', '的', '与', '及']:
                    matched.append(keyword)

    return matched


# ========================== Main Program Entry ==========================
def main():
    print("=" * 70)
    print("Sunshine College Exam Website Intangible Heritage Sports School & Major Data Crawler")
    print("=" * 70)

    try:
        # ---------- Step 1: Load matching keywords ----------
        keywords = read_keywords()
        print(f"Keyword preview (first 10 items): {keywords[:10]}")

        # ---------- Step 2: Fetch the list of offering universities ----------
        universities = get_university_list()

        if not universities:
            print("⚠️ No universities fetched, program terminated")
            return

        # ---------- Step 3: Iterate over universities and fetch detailed information ----------
        print("\n" + "=" * 70)
        print("🔍 Start iterating over universities to fetch course information")
        print("=" * 70)

        all_results = []
        total = len(universities)

        for idx, univ in enumerate(universities, 1):
            name = univ["name"]
            level = univ["level"]
            print(f"\n[{idx}/{total}] Processing: {name} ({level})")

            # Fetch detailed information
            univ = fetch_university_detail(univ)
            courses_text = univ.get("courses_raw", "")

            print(f"  📍 Location: {univ.get('location', 'Unknown')}")
            print(f"  🏫 Type: {univ.get('type', 'Unknown')}")
            print(f"  📖 Courses: {courses_text[:80]}..." if len(courses_text) > 80 else f"  📖 Courses: {courses_text}")

            # Match intangible heritage sports courses
            matched = match_courses(keywords, courses_text)

            if matched:
                print(f"  ✅ Matched {len(matched)} intangible heritage sports courses: {matched}")
                for course in matched:
                    all_results.append({
                        "对应非遗体育项目名称": course,
                        "院校名称": name,
                        "院校所在地": univ.get("location", ""),
                        "院校类型": univ.get("type", infer_university_type(name)),
                        "非遗传统体育课程名称": course,
                        "课程属性": "",  # Not provided on the Sunshine page
                        "课程等级": level,
                        "课程简介": f"{level}专业，开设课程包含：{courses_text[:100]}..." if courses_text else "",
                        "数据来源渠道": "阳光高考网",
                        "备注": f"专业层次：{level}"
                    })
            else:
                print(f"  ⚠️ No intangible heritage sports courses matched")

            # Request interval
            if idx < total:
                time.sleep(1.5)

        # ---------- Step 4: Deduplicate ----------
        if all_results:
            df_results = pd.DataFrame(all_results)
            # Deduplicate by university name + intangible heritage sports course name
            df_results = df_results.drop_duplicates(
                subset=["院校名称", "非遗传统体育课程名称"],
                keep="first"
            )
        else:
            df_results = pd.DataFrame(columns=[
                "对应非遗体育项目名称", "院校名称", "院校所在地", "院校类型",
                "非遗传统体育课程名称", "课程属性", "课程等级", "课程简介",
                "数据来源渠道", "备注"
            ])

        # ---------- Step 5: Export to excel ----------
        df_results.to_excel(OUTPUT_FILE, index=False)

        # ---------- Step 6: Print statistical report ----------
        print("\n" + "=" * 70)
        print("📊 Crawling‑task Completion Statistics")
        print("=" * 70)
        print(f"Total matching keywords: {len(keywords)}")
        print(f"Total processed universities: {total}")
        print(f"Total matched course records: {len(all_results)}")
        print(f"Deduplicated course records: {len(df_results)}")
        print(f"\n✅ Output file: {OUTPUT_FILE}")
        print("=" * 70)

        # Preview result data
        if not df_results.empty:
            print("\n📖 Result preview (first 10 entries):")
            print("-" * 70)
            preview_cols = ["对应非遗体育项目名称", "院校名称", "院校类型", "非遗传统体育课程名称"]
            print(df_results[preview_cols].head(10).to_string(index=False))
            print("-" * 70)

            # Count matched universities grouped by each keyword
            print("\n📊 Number of matched universities per intangible heritage sports project:")
            course_count = df_results.groupby("对应非遗体育项目名称").size().sort_values(ascending=False)
            for keyword, count in course_count.head(15).items():
                print(f"  {keyword}: {count} universities")
            if len(course_count) > 15:
                print(f"  ... {len(course_count) - 15} more projects omitted")
        else:
            print("\n⚠️ No course data matched")
            print("Potential reasons:")
            print("1. The major introduction pages on the Sunshine website do not list specific course names")
            print("2. Keywords and course names are expressed inconsistently")
            print("3. Access to more detailed course pages is required to obtain the data")

    except FileNotFoundError as e:
        print(f"❌ File‑related error: {e}")
    except ValueError as e:
        print(f"❌ Data‑validation error: {e}")
    except Exception as e:
        print(f"❌ Unexpected runtime error: {e}")
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()
